agora inbox for pgsql-hackers@postgresql.org  
help / color / mirror / Atom feed
[PATCH] Lock upgrade without deadlocks.
399+ messages / 2 participants
[nested] [flat]

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 399+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 399+ messages in thread

* [PATCH v1 2/2] Add per-backend lock statistics
@ 2026-06-03 13:05 Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
  0 siblings, 0 replies; 399+ messages in thread

From: Bertrand Drouvot @ 2026-06-03 13:05 UTC (permalink / raw)

This commit adds per-backend lock statistics, providing the same information as
pg_stat_lock, except that it is now possible to retrieve those stats (lock wait
counts, wait times, and fast-path exceeded count) on a per-backend basis.

This data can be retrieved with a new system function called
pg_stat_get_backend_lock(), that returns one tuple per lock type based on the PID
provided in input.  Like pg_stat_get_backend_io(), this is useful when joined
with pg_stat_activity to get a live picture of the locks behavior for each running
backend.

pgstat_flush_backend() gains a new flag value, able to control the flush of the
lock stats.

This commit relies mostly on the infrastructure provided by 9aea73fc61d4, that
has introduced backend statistics.

XXX: Bump catalog version.  A bump of PGSTAT_FILE_FORMAT_ID is not required,
as backend stats do not persist on disk.

Author: Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
Reviewed-by:
Discussion:
---
 doc/src/sgml/monitoring.sgml                | 19 ++++++
 src/backend/utils/activity/pgstat_backend.c | 70 +++++++++++++++++++++
 src/backend/utils/activity/pgstat_lock.c    |  4 ++
 src/backend/utils/adt/pgstatfuncs.c         | 29 ++++++++-
 src/include/catalog/pg_proc.dat             |  8 +++
 src/include/pgstat.h                        | 11 ++++
 src/include/utils/pgstat_internal.h         |  3 +-
 src/test/regress/expected/stats.out         | 12 ++++
 src/test/regress/sql/stats.sql              |  9 +++
 9 files changed, 162 insertions(+), 3 deletions(-)
  14.7% doc/src/sgml/
  38.6% src/backend/utils/activity/
  14.8% src/backend/utils/adt/
   8.7% src/include/catalog/
   9.9% src/include/
   6.8% src/test/regress/expected/
   6.1% src/test/regress/sql/

diff --git a/doc/src/sgml/monitoring.sgml b/doc/src/sgml/monitoring.sgml
index 08d5b824552..3936fb62a5d 100644
--- a/doc/src/sgml/monitoring.sgml
+++ b/doc/src/sgml/monitoring.sgml
@@ -5545,6 +5545,25 @@ description | Waiting for a newly initialized WAL file to reach durable storage
        </para></entry>
       </row>
 
+      <row>
+       <entry id="pg-stat-get-backend-lock" role="func_table_entry"><para role="func_signature">
+        <indexterm>
+         <primary>pg_stat_get_backend_lock</primary>
+        </indexterm>
+        <function>pg_stat_get_backend_lock</function> ( <type>integer</type> )
+        <returnvalue>setof record</returnvalue>
+       </para>
+       <para>
+        Returns lock statistics about the backend with the specified
+        process ID. The output fields are exactly the same as the ones in the
+        <structname>pg_stat_lock</structname> view.
+       </para>
+       <para>
+        The function does not return lock statistics for the checkpointer,
+        the background writer, the startup process and the autovacuum launcher.
+       </para></entry>
+      </row>
+
       <row>
        <entry role="func_table_entry"><para role="func_signature">
         <indexterm>
diff --git a/src/backend/utils/activity/pgstat_backend.c b/src/backend/utils/activity/pgstat_backend.c
index 73461c9bca5..297eda0a489 100644
--- a/src/backend/utils/activity/pgstat_backend.c
+++ b/src/backend/utils/activity/pgstat_backend.c
@@ -39,6 +39,7 @@
  */
 static PgStat_BackendPending PendingBackendStats;
 static bool backend_has_iostats = false;
+static bool backend_has_lockstats = false;
 
 /*
  * WAL usage counters saved from pgWalUsage at the previous call to
@@ -86,6 +87,37 @@ pgstat_count_backend_io_op(IOObject io_object, IOContext io_context,
 	pgstat_report_fixed = true;
 }
 
+/*
+ * Utility routines to report lock stats for backends, kept here to avoid
+ * exposing PendingBackendStats to the outside world.
+ */
+void
+pgstat_count_backend_lock_waits(uint8 locktag_type, long msecs)
+{
+	if (!pgstat_tracks_backend_bktype(MyBackendType))
+		return;
+
+	Assert(locktag_type <= LOCKTAG_LAST_TYPE);
+	PendingBackendStats.pending_lock.stats[locktag_type].waits++;
+	PendingBackendStats.pending_lock.stats[locktag_type].wait_time += (PgStat_Counter) msecs;
+
+	backend_has_lockstats = true;
+	pgstat_report_fixed = true;
+}
+
+void
+pgstat_count_backend_lock_fastpath_exceeded(uint8 locktag_type)
+{
+	if (!pgstat_tracks_backend_bktype(MyBackendType))
+		return;
+
+	Assert(locktag_type <= LOCKTAG_LAST_TYPE);
+	PendingBackendStats.pending_lock.stats[locktag_type].fastpath_exceeded++;
+
+	backend_has_lockstats = true;
+	pgstat_report_fixed = true;
+}
+
 /*
  * Returns statistics of a backend by proc number.
  */
@@ -262,6 +294,36 @@ pgstat_flush_backend_entry_wal(PgStat_EntryRef *entry_ref)
 	prevBackendWalUsage = pgWalUsage;
 }
 
+/*
+ * Flush out locally pending backend lock statistics.  Locking is managed
+ * by the caller.
+ */
+static void
+pgstat_flush_backend_entry_lock(PgStat_EntryRef *entry_ref)
+{
+	PgStatShared_Backend *shbackendent;
+	PgStat_PendingLock *bktype_shstats;
+
+	if (!backend_has_lockstats)
+		return;
+
+	shbackendent = (PgStatShared_Backend *) entry_ref->shared_stats;
+	bktype_shstats = &shbackendent->stats.lock_stats;
+
+	for (int i = 0; i <= LOCKTAG_LAST_TYPE; i++)
+	{
+#define LOCKSTAT_ACC(fld) \
+	(bktype_shstats->stats[i].fld += PendingBackendStats.pending_lock.stats[i].fld)
+		LOCKSTAT_ACC(waits);
+		LOCKSTAT_ACC(wait_time);
+		LOCKSTAT_ACC(fastpath_exceeded);
+#undef LOCKSTAT_ACC
+	}
+
+	MemSet(&PendingBackendStats.pending_lock, 0, sizeof(PgStat_PendingLock));
+	backend_has_lockstats = false;
+}
+
 /*
  * Flush out locally pending backend statistics
  *
@@ -286,6 +348,10 @@ pgstat_flush_backend(bool nowait, uint32 flags)
 		pgstat_backend_wal_have_pending())
 		has_pending_data = true;
 
+	/* Some lock data pending? */
+	if ((flags & PGSTAT_BACKEND_FLUSH_LOCK) && backend_has_lockstats)
+		has_pending_data = true;
+
 	if (!has_pending_data)
 		return false;
 
@@ -301,6 +367,9 @@ pgstat_flush_backend(bool nowait, uint32 flags)
 	if (flags & PGSTAT_BACKEND_FLUSH_WAL)
 		pgstat_flush_backend_entry_wal(entry_ref);
 
+	if (flags & PGSTAT_BACKEND_FLUSH_LOCK)
+		pgstat_flush_backend_entry_lock(entry_ref);
+
 	pgstat_unlock_entry(entry_ref);
 
 	return false;
@@ -339,6 +408,7 @@ pgstat_create_backend(ProcNumber procnum)
 
 	MemSet(&PendingBackendStats, 0, sizeof(PgStat_BackendPending));
 	backend_has_iostats = false;
+	backend_has_lockstats = false;
 
 	/*
 	 * Initialize prevBackendWalUsage with pgWalUsage so that
diff --git a/src/backend/utils/activity/pgstat_lock.c b/src/backend/utils/activity/pgstat_lock.c
index aec64f8fb4b..76116db3593 100644
--- a/src/backend/utils/activity/pgstat_lock.c
+++ b/src/backend/utils/activity/pgstat_lock.c
@@ -131,6 +131,8 @@ pgstat_count_lock_fastpath_exceeded(uint8 locktag_type)
 	PendingLockStats.stats[locktag_type].fastpath_exceeded++;
 	have_lockstats = true;
 	pgstat_report_fixed = true;
+
+	pgstat_count_backend_lock_fastpath_exceeded(locktag_type);
 }
 
 /*
@@ -147,4 +149,6 @@ pgstat_count_lock_waits(uint8 locktag_type, long msecs)
 	PendingLockStats.stats[locktag_type].wait_time += (PgStat_Counter) msecs;
 	have_lockstats = true;
 	pgstat_report_fixed = true;
+
+	pgstat_count_backend_lock_waits(locktag_type, msecs);
 }
diff --git a/src/backend/utils/adt/pgstatfuncs.c b/src/backend/utils/adt/pgstatfuncs.c
index 353607954ad..3f7c238e557 100644
--- a/src/backend/utils/adt/pgstatfuncs.c
+++ b/src/backend/utils/adt/pgstatfuncs.c
@@ -1740,8 +1740,8 @@ pg_stat_get_wal(PG_FUNCTION_ARGS)
 /*
  * pg_stat_lock_build_tuples
  *
- * Helper routine for pg_stat_get_lock(), filling a result tuplestore with one
- * tuple for each lock type.
+ * Helper routine for pg_stat_get_lock() and pg_stat_get_backend_lock(),
+ * filling a result tuplestore with one tuple for each lock type.
  */
 static void
 pg_stat_lock_build_tuples(ReturnSetInfo *rsinfo,
@@ -1788,6 +1788,31 @@ pg_stat_get_lock(PG_FUNCTION_ARGS)
 	return (Datum) 0;
 }
 
+/*
+ * Returns lock statistics for a backend with given PID.
+ */
+Datum
+pg_stat_get_backend_lock(PG_FUNCTION_ARGS)
+{
+	int			pid;
+	ReturnSetInfo *rsinfo;
+	PgStat_Backend *backend_stats;
+
+	InitMaterializedSRF(fcinfo, 0);
+	rsinfo = (ReturnSetInfo *) fcinfo->resultinfo;
+
+	pid = PG_GETARG_INT32(0);
+	backend_stats = pgstat_fetch_stat_backend_by_pid(pid, NULL);
+
+	if (!backend_stats)
+		return (Datum) 0;
+
+	pg_stat_lock_build_tuples(rsinfo, backend_stats->lock_stats.stats,
+							  backend_stats->stat_reset_timestamp);
+
+	return (Datum) 0;
+}
+
 /*
  * Returns statistics of SLRU caches.
  */
diff --git a/src/include/catalog/pg_proc.dat b/src/include/catalog/pg_proc.dat
index be157a5fbe9..87f7161527e 100644
--- a/src/include/catalog/pg_proc.dat
+++ b/src/include/catalog/pg_proc.dat
@@ -6092,6 +6092,14 @@
   proargmodes => '{i,o,o,o,o,o,o}',
   proargnames => '{backend_pid,wal_records,wal_fpi,wal_bytes,wal_fpi_bytes,wal_buffers_full,stats_reset}',
   prosrc => 'pg_stat_get_backend_wal' },
+{ oid => '9682', descr => 'statistics: backend lock statistics',
+  proname => 'pg_stat_get_backend_lock', prorows => '10', proretset => 't',
+  provolatile => 'v', proparallel => 'r', prorettype => 'record',
+  proargtypes => 'int4',
+  proallargtypes => '{int4,text,int8,int8,int8,timestamptz}',
+  proargmodes => '{i,o,o,o,o,o}',
+  proargnames => '{backend_pid,locktype,waits,wait_time,fastpath_exceeded,stats_reset}',
+  prosrc => 'pg_stat_get_backend_lock' },
 { oid => '6248', descr => 'statistics: information about WAL prefetching',
   proname => 'pg_stat_get_recovery_prefetch', prorows => '1', proretset => 't',
   provolatile => 'v', prorettype => 'record', proargtypes => '',
diff --git a/src/include/pgstat.h b/src/include/pgstat.h
index dfa2e837638..b5e69e6250e 100644
--- a/src/include/pgstat.h
+++ b/src/include/pgstat.h
@@ -523,6 +523,7 @@ typedef struct PgStat_Backend
 	TimestampTz stat_reset_timestamp;
 	PgStat_BktypeIO io_stats;
 	PgStat_WalCounters wal_counters;
+	PgStat_PendingLock lock_stats;
 } PgStat_Backend;
 
 /* ---------
@@ -535,6 +536,12 @@ typedef struct PgStat_BackendPending
 	 * Backend statistics store the same amount of IO data as PGSTAT_KIND_IO.
 	 */
 	PgStat_PendingIO pending_io;
+
+	/*
+	 * Backend statistics store the same amount of lock data as
+	 * PGSTAT_KIND_LOCK.
+	 */
+	PgStat_PendingLock pending_lock;
 } PgStat_BackendPending;
 
 /*
@@ -586,6 +593,10 @@ extern void pgstat_count_backend_io_op(IOObject io_object,
 									   IOContext io_context,
 									   IOOp io_op, uint32 cnt,
 									   uint64 bytes);
+
+/* used by pgstat_lock.c for lock stats tracked in backends */
+extern void pgstat_count_backend_lock_waits(uint8 locktag_type, long msecs);
+extern void pgstat_count_backend_lock_fastpath_exceeded(uint8 locktag_type);
 extern PgStat_Backend *pgstat_fetch_stat_backend(ProcNumber procNumber);
 extern PgStat_Backend *pgstat_fetch_stat_backend_by_pid(int pid,
 														BackendType *bktype);
diff --git a/src/include/utils/pgstat_internal.h b/src/include/utils/pgstat_internal.h
index fe463faaf63..b0788336ae3 100644
--- a/src/include/utils/pgstat_internal.h
+++ b/src/include/utils/pgstat_internal.h
@@ -705,7 +705,8 @@ extern void pgstat_archiver_snapshot_cb(void);
 /* flags for pgstat_flush_backend() */
 #define PGSTAT_BACKEND_FLUSH_IO		(1 << 0)	/* Flush I/O statistics */
 #define PGSTAT_BACKEND_FLUSH_WAL   (1 << 1) /* Flush WAL statistics */
-#define PGSTAT_BACKEND_FLUSH_ALL   (PGSTAT_BACKEND_FLUSH_IO | PGSTAT_BACKEND_FLUSH_WAL)
+#define PGSTAT_BACKEND_FLUSH_LOCK  (1 << 2) /* Flush lock statistics */
+#define PGSTAT_BACKEND_FLUSH_ALL   (PGSTAT_BACKEND_FLUSH_IO | PGSTAT_BACKEND_FLUSH_WAL | PGSTAT_BACKEND_FLUSH_LOCK)
 
 extern bool pgstat_flush_backend(bool nowait, uint32 flags);
 extern bool pgstat_backend_flush_cb(bool nowait);
diff --git a/src/test/regress/expected/stats.out b/src/test/regress/expected/stats.out
index bbb1db3c433..fa550676f83 100644
--- a/src/test/regress/expected/stats.out
+++ b/src/test/regress/expected/stats.out
@@ -2019,6 +2019,10 @@ BEGIN
 END;
 $$;
 SELECT fastpath_exceeded AS fastpath_exceeded_before FROM pg_stat_lock WHERE locktype = 'relation' \gset
+-- Test pg_stat_get_backend_lock()
+SELECT fastpath_exceeded AS backend_fastpath_exceeded_before
+  FROM pg_stat_get_backend_lock(pg_backend_pid())
+  WHERE locktype = 'relation' \gset
 -- Needs a lock on each partition
 SELECT count(*) FROM part_test;
  count 
@@ -2039,5 +2043,13 @@ SELECT fastpath_exceeded > :fastpath_exceeded_before FROM pg_stat_lock WHERE loc
  t
 (1 row)
 
+SELECT fastpath_exceeded > :backend_fastpath_exceeded_before
+  FROM pg_stat_get_backend_lock(pg_backend_pid())
+  WHERE locktype = 'relation';
+ ?column? 
+----------
+ t
+(1 row)
+
 DROP TABLE part_test;
 -- End of Stats Test
diff --git a/src/test/regress/sql/stats.sql b/src/test/regress/sql/stats.sql
index 610fd21fae4..f5683302a75 100644
--- a/src/test/regress/sql/stats.sql
+++ b/src/test/regress/sql/stats.sql
@@ -998,6 +998,11 @@ $$;
 
 SELECT fastpath_exceeded AS fastpath_exceeded_before FROM pg_stat_lock WHERE locktype = 'relation' \gset
 
+-- Test pg_stat_get_backend_lock()
+SELECT fastpath_exceeded AS backend_fastpath_exceeded_before
+  FROM pg_stat_get_backend_lock(pg_backend_pid())
+  WHERE locktype = 'relation' \gset
+
 -- Needs a lock on each partition
 SELECT count(*) FROM part_test;
 
@@ -1006,6 +1011,10 @@ SELECT pg_stat_force_next_flush();
 
 SELECT fastpath_exceeded > :fastpath_exceeded_before FROM pg_stat_lock WHERE locktype = 'relation';
 
+SELECT fastpath_exceeded > :backend_fastpath_exceeded_before
+  FROM pg_stat_get_backend_lock(pg_backend_pid())
+  WHERE locktype = 'relation';
+
 DROP TABLE part_test;
 
 -- End of Stats Test
-- 
2.34.1


--caEBMJLxMU8EntZL--






^ permalink  raw  reply  [nested|flat] 399+ messages in thread


end of thread, other threads:[~2026-06-03 13:05 UTC | newest]

Thread overview: 399+ messages (download: mbox mbox.gz follow: Atom feed)
-- links below jump to the message on this page --
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-06-03 13:05 [PATCH v1 2/2] Add per-backend lock statistics Bertrand Drouvot <bertranddrouvot.pg@gmail.com>

This inbox is served by agora; see mirroring instructions
for how to clone and mirror all data and code used for this inbox